agora inbox for [email protected]  
help / color / mirror / Atom feed
[PATCH v38 07/11] Add DISTINCT support for IVM
436+ messages / 2 participants
[nested] [flat]

* [PATCH v38 07/11] Add DISTINCT support for IVM
@ 2023-05-31 10:08  Yugo Nagata <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Yugo Nagata @ 2023-05-31 10:08 UTC (permalink / raw)

When IMMV is created with DISTINCT, multiplicity of tuples is
counted and stored in  "__ivm_count__" column, which is a hidden
column of IMMV.  The value in __ivm_count__ is updated when IMMV
is maintained incrementally. A tuple in IMMV can be removed if
and only if the count becomes zero.
---
 src/backend/commands/createas.c     | 141 +++++++++++++++++++++------
 src/backend/commands/indexcmds.c    |  40 ++++++++
 src/backend/commands/matview.c      | 145 ++++++++++++++++++++++++++--
 src/backend/commands/tablecmds.c    |   9 ++
 src/backend/nodes/outfuncs.c        |   1 +
 src/backend/nodes/readfuncs.c       |   1 +
 src/backend/parser/parse_relation.c |  18 +++-
 src/backend/rewrite/rewriteDefine.c |   3 +-
 src/include/commands/createas.h     |   2 +
 src/include/nodes/parsenodes.h      |   2 +
 10 files changed, 319 insertions(+), 43 deletions(-)

diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c
index a499688b79d..cd8db0059f9 100644
--- a/src/backend/commands/createas.c
+++ b/src/backend/commands/createas.c
@@ -55,6 +55,7 @@
 #include "parser/parser.h"
 #include "parser/parsetree.h"
 #include "parser/parse_clause.h"
+#include "parser/parse_func.h"
 #include "rewrite/rewriteHandler.h"
 #include "tcop/tcopprot.h"
 #include "utils/builtins.h"
@@ -307,6 +308,9 @@ ExecCreateTableAs(ParseState *pstate, CreateTableAsStmt *stmt,
 					 errhint("functions must be marked IMMUTABLE")));
 
 		check_ivm_restriction((Node *) query);
+
+		/* For IMMV, we need to rewrite matview query */
+		query = rewriteQueryForIMMV(query, into->colNames);
 	}
 
 	if (into->skipData)
@@ -416,6 +420,49 @@ ExecCreateTableAs(ParseState *pstate, CreateTableAsStmt *stmt,
 	return address;
 }
 
+/*
+ * rewriteQueryForIMMV -- rewrite view definition query for IMMV
+ *
+ * count(*) is added for counting distinct tuples in views.
+ */
+Query *
+rewriteQueryForIMMV(Query *query, List *colNames)
+{
+	Query *rewritten;
+
+	Node *node;
+	ParseState *pstate = make_parsestate(NULL);
+	FuncCall *fn;
+
+	rewritten = copyObject(query);
+	pstate->p_expr_kind = EXPR_KIND_SELECT_TARGET;
+
+	/*
+	 * Convert DISTINCT to GROUP BY and add count(*) for counting distinct
+	 * tuples in views.
+	 */
+	if (rewritten->distinctClause)
+	{
+		TargetEntry *tle;
+
+		rewritten->groupClause = transformDistinctClause(NULL, &rewritten->targetList, rewritten->sortClause, false);
+
+		fn = makeFuncCall(SystemFuncName("count"), NIL, COERCE_EXPLICIT_CALL, -1);
+		fn->agg_star = true;
+
+		node = ParseFuncOrColumn(pstate, fn->funcname, NIL, NULL, fn, false, -1);
+
+		tle = makeTargetEntry((Expr *) node,
+								list_length(rewritten->targetList) + 1,
+								pstrdup("__ivm_count__"),
+								false);
+		rewritten->targetList = lappend(rewritten->targetList, tle);
+		rewritten->hasAggs = true;
+	}
+
+	return rewritten;
+}
+
 /*
  * GetIntoRelEFlags --- compute executor flags needed for CREATE TABLE AS
  *
@@ -539,7 +586,8 @@ intorel_startup(DestReceiver *self, int operation, TupleDesc typeinfo)
 		ColumnDef  *col;
 		char	   *colname;
 
-		if (lc)
+		/* Don't override hidden columns added for IVM */
+		if (lc && !isIvmName(NameStr(attribute->attname)))
 		{
 			colname = strVal(lfirst(lc));
 			lc = lnext(into->colNames, lc);
@@ -943,10 +991,6 @@ check_ivm_restriction_walker(Node *node, void *context)
 					ereport(ERROR,
 							(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
 							 errmsg("LIMIT/OFFSET clause is not supported on incrementally maintainable materialized view")));
-				if (qry->distinctClause)
-					ereport(ERROR,
-							(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
-							 errmsg("DISTINCT is not supported on incrementally maintainable materialized view")));
 				if (qry->hasDistinctOn)
 					ereport(ERROR,
 							(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
@@ -1108,12 +1152,18 @@ CreateIndexOnIMMV(Query *query, Relation matviewRel)
 	char		idxname[NAMEDATALEN];
 	List	   *indexoidlist = RelationGetIndexList(matviewRel);
 	ListCell   *indexoidscan;
-	Bitmapset *key_attnos;
 
 	snprintf(idxname, sizeof(idxname), "%s_index", RelationGetRelationName(matviewRel));
 
 	index = makeNode(IndexStmt);
 
+	/*
+	 * We consider null values not distinct to make sure that views with DISTINCT
+	 * or GROUP BY don't contain multiple NULL rows when NULL is inserted to
+	 * a base table concurrently.
+	 */
+	index->nulls_not_distinct = true;
+
 	index->unique = true;
 	index->primary = false;
 	index->isconstraint = false;
@@ -1140,41 +1190,68 @@ CreateIndexOnIMMV(Query *query, Relation matviewRel)
 	index->concurrent = false;
 	index->if_not_exists = false;
 
-	/* create index on the base tables' primary key columns */
-	key_attnos = get_primary_key_attnos_from_query(query, &constraintList);
-	if (key_attnos)
+	if (query->distinctClause)
 	{
+		/* create unique constraint on all columns */
 		foreach(lc, query->targetList)
 		{
 			TargetEntry *tle = (TargetEntry *) lfirst(lc);
 			Form_pg_attribute attr = TupleDescAttr(matviewRel->rd_att, tle->resno - 1);
-
-			if (bms_is_member(tle->resno - FirstLowInvalidHeapAttributeNumber, key_attnos))
-			{
-				IndexElem  *iparam;
-
-				iparam = makeNode(IndexElem);
-				iparam->name = pstrdup(NameStr(attr->attname));
-				iparam->expr = NULL;
-				iparam->indexcolname = NULL;
-				iparam->collation = NIL;
-				iparam->opclass = NIL;
-				iparam->opclassopts = NIL;
-				iparam->ordering = SORTBY_DEFAULT;
-				iparam->nulls_ordering = SORTBY_NULLS_DEFAULT;
-				index->indexParams = lappend(index->indexParams, iparam);
-			}
+			IndexElem  *iparam;
+
+			iparam = makeNode(IndexElem);
+			iparam->name = pstrdup(NameStr(attr->attname));
+			iparam->expr = NULL;
+			iparam->indexcolname = NULL;
+			iparam->collation = NIL;
+			iparam->opclass = NIL;
+			iparam->opclassopts = NIL;
+			iparam->ordering = SORTBY_DEFAULT;
+			iparam->nulls_ordering = SORTBY_NULLS_DEFAULT;
+			index->indexParams = lappend(index->indexParams, iparam);
 		}
 	}
 	else
 	{
-		/* create no index, just notice that an appropriate index is necessary for efficient IVM */
-		ereport(NOTICE,
-				(errmsg("could not create an index on materialized view \"%s\" automatically",
-						RelationGetRelationName(matviewRel)),
-				 errdetail("This target list does not have all the primary key columns. "),
-				 errhint("Create an index on the materialized view for efficient incremental maintenance.")));
-		return;
+		Bitmapset *key_attnos;
+
+		/* create index on the base tables' primary key columns */
+		key_attnos = get_primary_key_attnos_from_query(query, &constraintList);
+		if (key_attnos)
+		{
+			foreach(lc, query->targetList)
+			{
+				TargetEntry *tle = (TargetEntry *) lfirst(lc);
+				Form_pg_attribute attr = TupleDescAttr(matviewRel->rd_att, tle->resno - 1);
+
+				if (bms_is_member(tle->resno - FirstLowInvalidHeapAttributeNumber, key_attnos))
+				{
+					IndexElem  *iparam;
+
+					iparam = makeNode(IndexElem);
+					iparam->name = pstrdup(NameStr(attr->attname));
+					iparam->expr = NULL;
+					iparam->indexcolname = NULL;
+					iparam->collation = NIL;
+					iparam->opclass = NIL;
+					iparam->opclassopts = NIL;
+					iparam->ordering = SORTBY_DEFAULT;
+					iparam->nulls_ordering = SORTBY_NULLS_DEFAULT;
+					index->indexParams = lappend(index->indexParams, iparam);
+				}
+			}
+		}
+		else
+		{
+			/* create no index, just notice that an appropriate index is necessary for efficient IVM */
+			ereport(NOTICE,
+					(errmsg("could not create an index on materialized view \"%s\" automatically",
+							RelationGetRelationName(matviewRel)),
+					 errdetail("This target list does not have all the primary key columns, "
+							   "or this view does not contain DISTINCT clause."),
+					 errhint("Create an index on the materialized view for efficient incremental maintenance.")));
+			return;
+		}
 	}
 
 	/* If we have a compatible index, we don't need to create another. */
diff --git a/src/backend/commands/indexcmds.c b/src/backend/commands/indexcmds.c
index 9ab74c8df0a..1f95e6b229f 100644
--- a/src/backend/commands/indexcmds.c
+++ b/src/backend/commands/indexcmds.c
@@ -41,6 +41,7 @@
 #include "commands/comment.h"
 #include "commands/defrem.h"
 #include "commands/event_trigger.h"
+#include "commands/matview.h"
 #include "commands/progress.h"
 #include "commands/tablecmds.h"
 #include "commands/tablespace.h"
@@ -1179,6 +1180,45 @@ DefineIndex(ParseState *pstate,
 	safe_index = indexInfo->ii_Expressions == NIL &&
 		indexInfo->ii_Predicate == NIL;
 
+	/*
+	 * We disallow unique indexes on IVM columns of IMMVs.
+	 */
+	if (RelationIsIVM(rel) && stmt->unique)
+	{
+		for (int i = 0; i < indexInfo->ii_NumIndexKeyAttrs; i++)
+		{
+			AttrNumber	attno = indexInfo->ii_IndexAttrNumbers[i];
+			if (attno > 0)
+			{
+				char *name = NameStr(TupleDescAttr(rel->rd_att, attno - 1)->attname);
+				if (name && isIvmName(name))
+					ereport(ERROR,
+							(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
+							 errmsg("unique index creation on IVM columns is not supported")));
+			}
+		}
+
+		if (indexInfo->ii_Expressions)
+		{
+			Bitmapset  *indexattrs = NULL;
+			int			varno = -1;
+
+			pull_varattnos((Node *) indexInfo->ii_Expressions, 1, &indexattrs);
+
+			while ((varno = bms_next_member(indexattrs, varno)) >= 0)
+			{
+				int attno = varno + FirstLowInvalidHeapAttributeNumber;
+				char *name = NameStr(TupleDescAttr(rel->rd_att, attno - 1)->attname);
+				if (name && isIvmName(name))
+					ereport(ERROR,
+							(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
+							 errmsg("unique index creation on IVM columns is not supported")));
+			}
+
+		}
+	}
+
+
 	/*
 	 * Report index creation if appropriate (delay this till after most of the
 	 * error checks)
diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c
index eaf39f80cd3..a2746ca9265 100644
--- a/src/backend/commands/matview.c
+++ b/src/backend/commands/matview.c
@@ -193,11 +193,15 @@ static Query *rewrite_query_for_postupdate_state(Query *query, MV_TriggerTable *
 
 static void apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *new_tuplestores,
 			TupleDesc tupdesc_old, TupleDesc tupdesc_new,
-			Query *query);
+			Query *query, bool use_count, char *count_colname);
 static void apply_old_delta(const char *matviewname, const char *deltaname_old,
 				List *keys);
+static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old,
+				List *keys, const char *count_colname);
 static void apply_new_delta(const char *matviewname, const char *deltaname_new,
 				StringInfo target_list);
+static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new,
+				List *keys, StringInfo target_list, const char* count_colname);
 static char *get_matching_condition_string(List *keys);
 static void generate_equal(StringInfo querybuf, Oid opttype,
 			   const char *leftop, const char *rightop);
@@ -346,6 +350,7 @@ RefreshMatViewByOid(Oid matviewOid, bool is_create, bool skipData,
 					QueryCompletion *qc)
 {
 	Relation	matviewRel;
+	Query	   *viewQuery;
 	Query	   *dataQuery;
 	Oid			tableSpace;
 	Oid			relowner;
@@ -394,7 +399,13 @@ RefreshMatViewByOid(Oid matviewOid, bool is_create, bool skipData,
 				 errmsg("%s options %s and %s cannot be used together",
 						"REFRESH", "CONCURRENTLY", "WITH NO DATA")));
 
-	dataQuery = get_matview_query(matviewRel);
+	viewQuery = get_matview_query(matviewRel);
+
+	/* For IMMV, we need to rewrite matview query */
+	if (!skipData && RelationIsIVM(matviewRel))
+		dataQuery = rewriteQueryForIMMV(viewQuery,NIL);
+	else
+		dataQuery = viewQuery;
 
 	/*
 	 * Check that there is a unique index with no WHERE clause on one or more
@@ -1688,6 +1699,13 @@ IVM_immediate_maintenance(PG_FUNCTION_ARGS)
 			int	rte_index = lfirst_int(lc2);
 			TupleDesc		tupdesc_old;
 			TupleDesc		tupdesc_new;
+			bool	use_count = false;
+			char   *count_colname = NULL;
+
+			count_colname = pstrdup("__ivm_count__");
+
+			if (query->distinctClause)
+				use_count = true;
 
 			/* calculate delta tables */
 			calc_delta(table, rte_index, rewritten, dest_old, dest_new,
@@ -1700,7 +1718,8 @@ IVM_immediate_maintenance(PG_FUNCTION_ARGS)
 			{
 				/* apply the delta tables to the materialized view */
 				apply_delta(matviewOid, old_tuplestore, new_tuplestore,
-							tupdesc_old, tupdesc_new, query);
+							tupdesc_old, tupdesc_new, query, use_count,
+							count_colname);
 			}
 			PG_CATCH();
 			{
@@ -2238,7 +2257,7 @@ rewrite_query_for_postupdate_state(Query *query, MV_TriggerTable *table, int rte
 static void
 apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *new_tuplestores,
 			TupleDesc tupdesc_old, TupleDesc tupdesc_new,
-			Query *query)
+			Query *query, bool use_count, char *count_colname)
 {
 	StringInfoData querybuf;
 	StringInfoData target_list_buf;
@@ -2314,7 +2333,12 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n
 		if (rc != SPI_OK_REL_REGISTER)
 			elog(ERROR, "SPI_register failed");
 
-		apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys);
+		if (use_count)
+			/* apply old delta and get rows to be recalculated */
+			apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME,
+									   keys, count_colname);
+		else
+			apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys);
 
 	}
 	/* For tuple insertion */
@@ -2336,7 +2360,11 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n
 			elog(ERROR, "SPI_register failed");
 
 		/* apply new delta */
-		apply_new_delta(matviewname, NEW_DELTA_ENRNAME, &target_list_buf);
+		if (use_count)
+			apply_new_delta_with_count(matviewname, NEW_DELTA_ENRNAME,
+								keys, &target_list_buf, count_colname);
+		else
+			apply_new_delta(matviewname, NEW_DELTA_ENRNAME, &target_list_buf);
 	}
 
 	/* We're done maintaining the materialized view. */
@@ -2349,6 +2377,51 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n
 		elog(ERROR, "SPI_finish failed");
 }
 
+/*
+ * apply_old_delta_with_count
+ *
+ * Execute a query for applying a delta table given by deltname_old
+ * which contains tuples to be deleted from to a materialized view given by
+ * matviewname.  This is used when counting is required, that is, the view
+ * has aggregate or distinct.
+ */
+static void
+apply_old_delta_with_count(const char *matviewname, const char *deltaname_old,
+				List *keys, const char *count_colname)
+{
+	StringInfoData	querybuf;
+	char   *match_cond;
+
+	/* build WHERE condition for searching tuples to be deleted */
+	match_cond = get_matching_condition_string(keys);
+
+	/* Search for matching tuples from the view and update or delete if found. */
+	initStringInfo(&querybuf);
+	appendStringInfo(&querybuf,
+					"WITH t AS ("			/* collecting tid of target tuples in the view */
+						"SELECT diff.%s, "			/* count column */
+								"(diff.%s OPERATOR(pg_catalog.=) mv.%s) AS for_dlt, "
+								"mv.ctid "
+						"FROM %s AS mv, %s AS diff "
+						"WHERE %s"					/* tuple matching condition */
+					"), updt AS ("			/* update a tuple if this is not to be deleted */
+						"UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s "
+						"FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt "
+					")"
+					/* delete a tuple if this is to be deleted */
+					"DELETE FROM %s AS mv USING t "
+					"WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt",
+					count_colname,
+					count_colname, count_colname,
+					matviewname, deltaname_old,
+					match_cond,
+					matviewname, count_colname, count_colname, count_colname,
+					matviewname);
+
+	if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE)
+		elog(ERROR, "SPI_exec failed: %s", querybuf.data);
+}
+
 /*
  * apply_old_delta
  *
@@ -2398,6 +2471,66 @@ apply_old_delta(const char *matviewname, const char *deltaname_old,
 		elog(ERROR, "SPI_exec failed: %s", querybuf.data);
 }
 
+/*
+ * apply_new_delta_with_count
+ *
+ * Execute a query for applying a delta table given by deltname_new
+ * which contains tuples to be inserted into a materialized view given by
+ * matviewname.  This is used when counting is required, that is, the view
+ * has aggregate or distinct. Also, when a table in EXISTS sub queries
+ * is modified.
+ */
+static void
+apply_new_delta_with_count(const char *matviewname, const char* deltaname_new,
+				List *keys, StringInfo target_list, const char* count_colname)
+{
+	StringInfoData	querybuf;
+	StringInfoData	returning_keys;
+	ListCell	*lc;
+	char	*match_cond = "";
+
+	/* build WHERE condition for searching tuples to be updated */
+	match_cond = get_matching_condition_string(keys);
+
+	/* build string of keys list */
+	initStringInfo(&returning_keys);
+	if (keys)
+	{
+		foreach (lc, keys)
+		{
+			Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc);
+			char   *resname = NameStr(attr->attname);
+			appendStringInfo(&returning_keys, "%s", quote_qualified_identifier("mv", resname));
+			if (lnext(keys, lc))
+				appendStringInfo(&returning_keys, ", ");
+		}
+	}
+	else
+		appendStringInfo(&returning_keys, "NULL");
+
+	/* Search for matching tuples from the view and update if found or insert if not. */
+	initStringInfo(&querybuf);
+	appendStringInfo(&querybuf,
+					"WITH updt AS ("		/* update a tuple if this exists in the view */
+						"UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.+) diff.%s "
+						"FROM %s AS diff "
+						"WHERE %s "					/* tuple matching condition */
+						"RETURNING %s"				/* returning keys of updated tuples */
+					") INSERT INTO %s (%s) "	/* insert a new tuple if this doesn't exist */
+						"SELECT %s FROM %s AS diff "
+						"WHERE NOT EXISTS (SELECT 1 FROM updt AS mv WHERE %s);",
+					matviewname, count_colname, count_colname, count_colname,
+					deltaname_new,
+					match_cond,
+					returning_keys.data,
+					matviewname, target_list->data,
+					target_list->data, deltaname_new,
+					match_cond);
+
+	if (SPI_exec(querybuf.data, 0) != SPI_OK_INSERT)
+		elog(ERROR, "SPI_exec failed: %s", querybuf.data);
+}
+
 /*
  * apply_new_delta
  *
diff --git a/src/backend/commands/tablecmds.c b/src/backend/commands/tablecmds.c
index 920b62d254d..f62765f0a89 100644
--- a/src/backend/commands/tablecmds.c
+++ b/src/backend/commands/tablecmds.c
@@ -60,6 +60,7 @@
 #include "catalog/toasting.h"
 #include "commands/comment.h"
 #include "commands/defrem.h"
+#include "commands/matview.h"
 #include "commands/event_trigger.h"
 #include "commands/extension.h"
 #include "commands/matview.h"
@@ -3930,6 +3931,14 @@ renameatt_internal(Oid myrelid,
 	targetrelation = relation_open(myrelid, AccessExclusiveLock);
 	renameatt_check(myrelid, RelationGetForm(targetrelation), recursing);
 
+	/*
+	 * Don't rename IVM columns.
+	 */
+	if (RelationIsIVM(targetrelation) && isIvmName(oldattname))
+		ereport(ERROR,
+				(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
+				 errmsg("IVM column can not be renamed")));
+
 	/*
 	 * if the 'recurse' flag is set then we are supposed to rename this
 	 * attribute in all classes that inherit from 'relname' (as well as in
diff --git a/src/backend/nodes/outfuncs.c b/src/backend/nodes/outfuncs.c
index 953c5797c5d..0f228eca259 100644
--- a/src/backend/nodes/outfuncs.c
+++ b/src/backend/nodes/outfuncs.c
@@ -516,6 +516,7 @@ _outRangeTblEntry(StringInfo str, const RangeTblEntry *node)
 			WRITE_INT_FIELD(rellockmode);
 			WRITE_UINT_FIELD(perminfoindex);
 			WRITE_NODE_FIELD(tablesample);
+			WRITE_BOOL_FIELD(relisivm);
 			break;
 		case RTE_SUBQUERY:
 			WRITE_NODE_FIELD(subquery);
diff --git a/src/backend/nodes/readfuncs.c b/src/backend/nodes/readfuncs.c
index b6b2ce6c792..dba4d8ff2de 100644
--- a/src/backend/nodes/readfuncs.c
+++ b/src/backend/nodes/readfuncs.c
@@ -365,6 +365,7 @@ _readRangeTblEntry(void)
 			READ_INT_FIELD(rellockmode);
 			READ_UINT_FIELD(perminfoindex);
 			READ_NODE_FIELD(tablesample);
+			READ_BOOL_FIELD(relisivm);
 			break;
 		case RTE_SUBQUERY:
 			READ_NODE_FIELD(subquery);
diff --git a/src/backend/parser/parse_relation.c b/src/backend/parser/parse_relation.c
index ced210cd206..5392e9f15b2 100644
--- a/src/backend/parser/parse_relation.c
+++ b/src/backend/parser/parse_relation.c
@@ -33,6 +33,7 @@
 #include "utils/lsyscache.h"
 #include "utils/syscache.h"
 #include "utils/varlena.h"
+#include "commands/matview.h"
 
 
 /*
@@ -96,7 +97,7 @@ static void expandTupleDesc(TupleDesc tupdesc, Alias *eref,
 							int rtindex, int sublevels_up,
 							VarReturningType returning_type,
 							int location, bool include_dropped,
-							List **colnames, List **colvars);
+							List **colnames, List **colvars, bool is_ivm);
 static int	specialAttNum(const char *attname);
 static bool rte_visible_if_lateral(ParseState *pstate, RangeTblEntry *rte);
 static bool rte_visible_if_qualified(ParseState *pstate, RangeTblEntry *rte);
@@ -1544,6 +1545,7 @@ addRangeTableEntry(ParseState *pstate,
 	rte->inh = inh;
 	rte->relkind = rel->rd_rel->relkind;
 	rte->rellockmode = lockmode;
+	rte->relisivm = rel->rd_rel->relisivm;
 
 	/*
 	 * Build the list of effective column names using user-supplied aliases
@@ -1625,6 +1627,7 @@ addRangeTableEntryForRelation(ParseState *pstate,
 	rte->inh = inh;
 	rte->relkind = rel->rd_rel->relkind;
 	rte->rellockmode = lockmode;
+	rte->relisivm = rel->rd_rel->relisivm;
 
 	/*
 	 * Build the list of effective column names using user-supplied aliases
@@ -2967,7 +2970,7 @@ expandRTE(RangeTblEntry *rte, int rtindex, int sublevels_up,
 										rtfunc->funccolcount, atts_done,
 										rtindex, sublevels_up,
 										returning_type, location,
-										include_dropped, colnames, colvars);
+										include_dropped, colnames, colvars, false);
 					}
 					else if (functypclass == TYPEFUNC_SCALAR)
 					{
@@ -3243,7 +3246,7 @@ expandRelation(Oid relid, Alias *eref, int rtindex, int sublevels_up,
 	expandTupleDesc(rel->rd_att, eref, rel->rd_att->natts, 0,
 					rtindex, sublevels_up, returning_type,
 					location, include_dropped,
-					colnames, colvars);
+					colnames, colvars, RelationIsIVM(rel));
 	relation_close(rel, AccessShareLock);
 }
 
@@ -3261,7 +3264,7 @@ expandTupleDesc(TupleDesc tupdesc, Alias *eref, int count, int offset,
 				int rtindex, int sublevels_up,
 				VarReturningType returning_type,
 				int location, bool include_dropped,
-				List **colnames, List **colvars)
+				List **colnames, List **colvars, bool is_ivm)
 {
 	ListCell   *aliascell;
 	int			varattno;
@@ -3274,6 +3277,9 @@ expandTupleDesc(TupleDesc tupdesc, Alias *eref, int count, int offset,
 	{
 		Form_pg_attribute attr = TupleDescAttr(tupdesc, varattno);
 
+		if (is_ivm && isIvmName(NameStr(attr->attname)) && !MatViewIncrementalMaintenanceIsEnabled())
+			continue;
+
 		if (attr->attisdropped)
 		{
 			if (include_dropped)
@@ -3438,6 +3444,10 @@ expandNSItemAttrs(ParseState *pstate, ParseNamespaceItem *nsitem,
 		Var		   *varnode = (Var *) lfirst(var);
 		TargetEntry *te;
 
+		/* if transform * into columnlist with IMMV, remove IVM columns */
+		if (rte->relisivm && isIvmName(label) && !MatViewIncrementalMaintenanceIsEnabled())
+			continue;
+
 		te = makeTargetEntry((Expr *) varnode,
 							 (AttrNumber) pstate->p_next_resno++,
 							 label,
diff --git a/src/backend/rewrite/rewriteDefine.c b/src/backend/rewrite/rewriteDefine.c
index 6a223fbeaa4..74ac631299c 100644
--- a/src/backend/rewrite/rewriteDefine.c
+++ b/src/backend/rewrite/rewriteDefine.c
@@ -614,7 +614,8 @@ checkRuleResultList(List *targetList, TupleDesc resultDesc, bool isSelect,
 														attr->atttypmod))));
 	}
 
-	if (i != resultDesc->natts)
+	/* No check for materialized views since this could have special columns for IVM */
+	if ((!isSelect || requireColumnNameMatch) && i != resultDesc->natts)
 		ereport(ERROR,
 				(errcode(ERRCODE_INVALID_OBJECT_DEFINITION),
 				 isSelect ?
diff --git a/src/include/commands/createas.h b/src/include/commands/createas.h
index c286ebcd70e..bfd0249b10d 100644
--- a/src/include/commands/createas.h
+++ b/src/include/commands/createas.h
@@ -29,6 +29,8 @@ extern ObjectAddress ExecCreateTableAs(ParseState *pstate, CreateTableAsStmt *st
 extern void CreateIvmTriggersOnBaseTables(Query *qry, Oid matviewOid);
 extern void CreateIndexOnIMMV(Query *query, Relation matviewRel);
 
+extern Query *rewriteQueryForIMMV(Query *query, List *colNames);
+
 extern int	GetIntoRelEFlags(IntoClause *intoClause);
 
 extern DestReceiver *CreateIntoRelDestReceiver(IntoClause *intoClause);
diff --git a/src/include/nodes/parsenodes.h b/src/include/nodes/parsenodes.h
index 4133c404a6b..cbcef860061 100644
--- a/src/include/nodes/parsenodes.h
+++ b/src/include/nodes/parsenodes.h
@@ -1209,6 +1209,8 @@ typedef struct RangeTblEntry
 	Index		perminfoindex pg_node_attr(query_jumble_ignore);
 	/* sampling info, or NULL */
 	struct TableSampleClause *tablesample;
+	/* incrementally maintainable materialized view? */
+	bool		relisivm;
 
 	/*
 	 * Fields valid for a subquery RTE (else NULL):
-- 
2.43.0


--Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ
Content-Type: text/x-diff;
 name="v38-0006-Add-Incremental-View-Maintenance-support.patch"
Content-Disposition: attachment;
 filename="v38-0006-Add-Incremental-View-Maintenance-support.patch"
Content-Transfer-Encoding: 7bit



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread

* [PATCH v3 1/2] ci: Improve ccache handling
@ 2026-06-05 15:58  Andres Freund <[email protected]>
  0 siblings, 0 replies; 436+ messages in thread

From: Andres Freund @ 2026-06-05 15:58 UTC (permalink / raw)

There previously were a number of issues:

- We'd upload the cache even if we already had a high hit rate. That means we
  churn through the available cache space very quickly.

  For this we now check if the cache hit ratio is already high, and skip
  uploading a new cache in that case.

- We'd generate per-branch caches, even if master's already would suffice,
  because the branch doesn't change much

  This is solved indirectly by the above.

- The cache key allowed prefix matches based on the branch,
  e.g. master-pending would always use master's branch

  Replace the cache key element separator of - with :, which is not a valid
  part of a branch name.

- When rebasing a feature branch, we'd start with just that branch's cache,
  rather than also having the newer cache of master available

  This is solved by downloading by master's and the feature branch's cache,
  simply overlaying both. That's possible because ccache is content addressed.

- The size of a cache would increase to the max, even though there likely will
  be no benefit from old cache entries.

  Address this by explicitly evicting old data and also recompressing the
  cache before uploading it.

In my testing this utilizes the available cache space (10GB for personal
accounts) much more effictively than before.

The not entirely trivial determination of whether it's worth uploading a cache
entry is moved to a python script.  I first had it as shell, but that gets
awkward.  This way it'd also be more viable to use ccache for msvc at some
point.

The per-job redundancies are a bit annoying. There's a way around that, by
using composite actions, but I think that might be harder to understand,
without all that much of an improvement.
---
 .github/workflows/pg-ci.yml       |  94 ++++++++++++++++++++++------
 src/tools/ci/gha_ccache_decide.py | 100 ++++++++++++++++++++++++++++++
 2 files changed, 176 insertions(+), 18 deletions(-)
 create mode 100644 src/tools/ci/gha_ccache_decide.py

diff --git a/.github/workflows/pg-ci.yml b/.github/workflows/pg-ci.yml
index 8560e9389f6..86dc47de8db 100644
--- a/.github/workflows/pg-ci.yml
+++ b/.github/workflows/pg-ci.yml
@@ -130,6 +130,22 @@ env:
   # commit-message directive parsed in the `setup` job below.
   CI_OS_ONLY_JOBS: "linux macos windows mingw compilerwarnings sanitycheck"
 
+  ###
+  # A few variables to make expressions later on shorter
+  ###
+
+  ON_DEFAULT_BRANCH: ${{github.event.repository.default_branch == github.ref_name }}
+
+  # Note that we need to be careful to use a separator that can't be in branch
+  # names, otherwise e.g. caches for 'master' might be restored on the
+  # 'master-pending' branch.
+  CACHE_PREFIX_DEFAULT: >-
+    :${{ github.job }}:${{ github.event.repository.default_branch }}:
+  CACHE_PREFIX_BRANCH: >-
+    :${{ github.job }}:${{ github.ref_name }}:
+  CACHE_SUFFIX: >-
+    ${{ github.run_id }}:${{ github.run_attempt }}
+
 
 jobs:
 
@@ -277,16 +293,30 @@ jobs:
         with:
           fetch-depth: ${{ env.CLONE_DEPTH }}
 
-      - &ccache_restore_step
-        name: Restore ccache
-        id: ccache_restore
+      # We restore both the ccache from the default branch (typically master),
+      # and from the current branch. This will often allow feature branches to
+      # start out with a high cache hit ratio.
+      #
+      # With ccache it turns out to work to just restore two caches into the
+      # same directory, as it's basically a content addressed store. Stats
+      # could be corrupted, but we zero them out anyway.
+      - &ccache_restore_default_step
+        name: "ccache: Restore for default branch ${{github.event.repository.default_branch}}"
+        if: ${{ env.ON_DEFAULT_BRANCH == 'false' }}
         uses: actions/cache/restore@v5
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ccache-${{ github.job }}-${{ github.ref_name }}-${{ github.run_id }}-${{ github.run_attempt }}
-          restore-keys: |
-            ccache-${{ github.job }}-${{ github.ref_name }}-
-            ccache-${{ github.job }}-
+          key: ccache${{env.CACHE_PREFIX_DEFAULT}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_DEFAULT}}
+
+      - &ccache_restore_branch_step
+        name: "ccache: Restore for branch ${{ github.ref_name }}"
+        id: ccache-restore-branch
+        uses: actions/cache/restore@v5
+        with:
+          path: ${{ env.CCACHE_DIR }}
+          key: ccache${{env.CACHE_PREFIX_BRANCH}}${{env.CACHE_SUFFIX}}
+          restore-keys: ccache${{env.CACHE_PREFIX_BRANCH}}
 
       - &linux_prepare_workspace_step
         name: Prepare workspace
@@ -325,15 +355,30 @@ jobs:
           ninja -C build -j${{env.BUILD_JOBS}} ${{env.MBUILD_TARGET}}
           ninja -C build -t missingdeps
 
-      # TODO: As long as we use per-run ccache caches, we should probably add
-      # a step that checks if there is sufficient new content to warrant
-      # saving the new cache.
+      # Decide if it's worth uploading a new version of the ccache cache. If
+      # we always do so unconditionally, we'd very quickly go through the
+      # allowed cache space. Instead we check if the hit rate is high enough
+      # already for that not to be worth it.
+      - &ccache_decide_save_step
+        name: "ccache: Decide if cache should be uploaded"
+        id: ccache-pre-save
+        # [Decide to] store the cache whenever the cache was set up, so that
+        # incrementally addressing compiler errors/warnings doesn't have to
+        # start from scratch.
+        if: |
+          always() &&
+          steps.ccache-restore-branch.conclusion == 'success'
+        run: python3 src/tools/ci/gha_ccache_decide.py
+
       - &ccache_save_step
-        name: Save ccache
+        name: "ccache: Upload cache"
         uses: actions/cache/save@v5
+        if: |
+          always() &&
+          steps.ccache-pre-save.outputs.should_save == 'true'
         with:
           path: ${{ env.CCACHE_DIR }}
-          key: ${{ steps.ccache_restore.outputs.cache-primary-key }}
+          key: ${{ steps.ccache-restore-branch.outputs.cache-primary-key }}
 
       # Run a minimal set of tests. The main regression tests take too long
       # for this purpose. For now this is a random quick pg_regress style
@@ -448,7 +493,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -467,6 +513,7 @@ jobs:
         run: |
           make -s -j${BUILD_JOBS} world-bin
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -508,7 +555,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -527,6 +575,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -596,7 +645,8 @@ jobs:
 
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
       - *linux_prepare_workspace_step
 
       - name: Configure
@@ -613,6 +663,7 @@ jobs:
         shell: *su_postgres_shell
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -682,7 +733,6 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
 
       - name: Setup core files
         run: |
@@ -745,6 +795,9 @@ jobs:
           path: ${{ env.MACPORTS_CACHE }}
           key: ${{ steps.mp-key.outputs.key }}
 
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
+
       - name: Configure
         env:
           PKG_CONFIG_PATH: /opt/local/lib/pkgconfig/
@@ -762,6 +815,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1062,7 +1116,8 @@ jobs:
         shell: cmd
         run: mkdir ${{env.PG_REGRESS_SOCK_DIR}}
 
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Configure
         run: |
@@ -1077,6 +1132,7 @@ jobs:
       - name: Build
         run: *ninja_build_cmd
 
+      - *ccache_decide_save_step
       - *ccache_save_step
 
       - name: Test world
@@ -1118,7 +1174,8 @@ jobs:
     steps:
       - *nix_sysinfo_step
       - *checkout_step
-      - *ccache_restore_step
+      - *ccache_restore_default_step
+      - *ccache_restore_branch_step
 
       - name: Setup workspace
         run: |
@@ -1213,5 +1270,6 @@ jobs:
             headerscheck cpluspluscheck \
             EXTRAFLAGS='-fmax-errors=10'
 
+      - *ccache_decide_save_step
       - *ccache_save_step
       - *upload_logs_step
diff --git a/src/tools/ci/gha_ccache_decide.py b/src/tools/ci/gha_ccache_decide.py
new file mode 100644
index 00000000000..920f7bf9685
--- /dev/null
+++ b/src/tools/ci/gha_ccache_decide.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+
+import os
+import re
+import shutil
+import subprocess
+
+def run(cmd, check=True):
+    return subprocess.run(
+        cmd,
+        check=check,
+        text=True,
+        stdout=subprocess.PIPE,
+        stderr=subprocess.STDOUT,
+    ).stdout
+
+def parse_ccache_stats():
+    out = run(["ccache", "--print-stats"])
+    hits = 0
+    misses = 0
+
+    for line in out.splitlines():
+        line = line.strip()
+        m = re.match(r"^local_storage_hit\s+(\d+)$", line)
+        if m:
+            hits = int(m.group(1))
+            continue
+        m = re.match(r"^local_storage_miss\s+(\d+)$", line)
+        if m:
+            misses = int(m.group(1))
+            continue
+
+    return hits, misses
+
+def append_github_output(key, value):
+    output_path = os.environ["GITHUB_OUTPUT"]
+    with open(output_path, "a", encoding="utf-8") as f:
+        f.write(f"{key}={value}\n")
+
+def main():
+    on_default_branch = os.environ["ON_DEFAULT_BRANCH"] == "true"
+    ccache_dir = os.environ["CCACHE_DIR"]
+
+    # Decide the target hit percentage below which we decide to upload a new
+    # cache. On non-default branches a few misses aren't that bad. But, as the
+    # caches of the default branch are shared with all branches, it's worth
+    # aiming for a higher ratio there.
+    target_rate = 95 if on_default_branch else 80
+
+    # Log ccache stats, useful for more in-depth understanding. The avoid it
+    # swamping the output, collapse it in a group.
+    print("::group::ccache_stats")
+    print(run(["ccache", "-s", "-vv"]))
+    print("::endgroup::")
+
+    # compute cache hit ratio
+    hits, misses = parse_ccache_stats()
+    total = hits + misses
+    hit_pct = int(( hits / total) * 100) if total > 0 else 100
+
+    print(f"hits: {hits}, misses: {misses}, hit_pct: {hit_pct}, target rate: {target_rate}")
+
+    # If there were either barely any misses, or the cache hit ratio was high,
+    # there no point in generating a new cache entry. We have limited cache
+    # space.
+    should_save = misses > 10 and hit_pct < target_rate
+
+    append_github_output("should_save", str(should_save).lower())
+
+    if not should_save:
+        print(f"hit rate {hit_pct} is above target of {target_rate}, skip creating new cache entry")
+        return 0
+
+    print(f"hit rate {hit_pct} is below target of {target_rate}, create new cache entry")
+
+    # It's not worth persisting old cache entries (e.g. from before a
+    # change to a central header, or from the default branch if this
+    # branch differs a lot). Therefore evict ccache entries that are a
+    # bit older. The cutoff here is fairly arbitrary, it could
+    # probably be improved.
+    print("::group::ccache_shrink")
+    print(run(["ccache", "--evict-older-than", f"{45*60}s"]))
+    print(run(["ccache", "-X", "10"]))
+
+    # Don't store ccache stats , otherwise we'd need to reset the cache access
+    # data after restoring the cache in the next run, to be able to get the
+    # hit ratio of the CI run.
+    print(run(["ccache", "-z"]))
+    print("::endgroup::")
+
+    # Before continuing, try to kill all ccache instances, otherwise
+    # it's possible that on cancellations there is still running
+    # ccaches that cause the upload to fail.
+    if shutil.which("killall"):
+        print(run(["killall", "ccache"], check=False))
+
+    return 0
+
+if __name__ == "__main__":
+    exit(main())
-- 
2.54.0.380.gc69baaf57b


--zzyr2ozda5zjkkci
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment; filename="v3-0002-ci-fewer-tests.patch"



^ permalink  raw  reply  [nested|flat] 436+ messages in thread


end of thread, other threads:[~2026-06-05 15:58 UTC | newest]

Thread overview: 436+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2023-05-31 10:08 [PATCH v38 07/11] Add DISTINCT support for IVM Yugo Nagata <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>
2026-06-05 15:58 [PATCH v3 1/2] ci: Improve ccache handling Andres Freund <[email protected]>

This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox