pgpool2.git / summary / log / commit / refs
commit 2efc59eeeff04ca2bb986c9591a249d8b64188f2
Author: Tatsuo Ishii <ishii@sraoss.co.jp>
Date: Sat Aug 19 06:44:02 2023 +0000
Feature: allow to set delay_threshold_by_time in milliseconds.
Previously it was allowed only in seconds. Also put some
refactoring. Create new function "check_replication_delay" which
checks the replication delay and returns 0, -1 or -2, depending on "no
delay", "delayed (delay_threshold_by_time)" or "delayed
(delay_threshold)" accordingly. This should simplify the lengthy
if-statement to check the replication delay.
Discussion: https://www.pgpool.net/pipermail/pgpool-hackers/2023-August/004372.html
doc.ja/src/sgml/stream-check.sgml | 11 +++--
doc/src/sgml/stream-check.sgml | 7 +--
src/config/pool_config_variables.c | 2 +-
src/context/pool_query_context.c | 10 +---
src/include/protocol/pool_pg_utils.h | 1 +
src/protocol/pool_pg_utils.c | 53 ++++++++++++++++++----
src/sample/pgpool.conf.sample-stream | 4 +-
src/streaming_replication/pool_worker_child.c | 4 +-
.../tests/033.prefer_lower_standby_delay/test.sh | 4 +-
9 files changed, 64 insertions(+), 32 deletions(-)
diff --git a/doc.ja/src/sgml/stream-check.sgml b/doc.ja/src/sgml/stream-check.sgml
index dcb9846e4..396adf397 100644
--- a/doc.ja/src/sgml/stream-check.sgml
+++ b/doc.ja/src/sgml/stream-check.sgml
@@ -270,9 +270,10 @@
<para>
<!--
- Specifies the maximum tolerance level of replication delay in
- seconds on the standby server against the primary server. If the
- specified value is greater than
+ Specifies the maximum tolerance level of replication delay
+ on the standby server against the primary server.
+ If this value is specified without units, it is taken as milliseconds.
+ If the specified value is greater than
0, <xref linkend="guc-delay-threshold"> is ignored. If the delay
exceeds this configured level,
<productname>Pgpool-II</productname> stops sending the <acronym>
@@ -283,7 +284,9 @@
This delay threshold check is performed every <xref linkend="guc-sr-check-period">.
Default is 0.
-->
- プライマリサーバに対するスタンバイサーバのレプリケーション遅延の許容度を秒単位で指定します。
+ プライマリサーバに対するスタンバイサーバのレプリケーション遅延の許容時間を指定します。
+ この値が単位無しで指定された場合は、マイクロ秒単位であると見なします。
+ 0よりも大きい値が指定されると、<xref linkend="guc-delay-threshold">は無視されます。
<productname>Pgpool-II</productname>は、スタンバイサーバの遅延がこの設定レベルを超えた場合には、 <xref linkend="guc-load-balance-mode">が有効であっても、プライマリに追いつくまでそのスタンバイノードには<acronym>SELECT</acronym>クエリを送信せず、全てプライマリサーバに送るようにします。
このパラメータが0の場合は、遅延のチェックを行ないません。
この遅延閾値のチェックは<xref linkend="guc-sr-check-period">毎に行われます。
diff --git a/doc/src/sgml/stream-check.sgml b/doc/src/sgml/stream-check.sgml
index 25b412353..c86c9322c 100644
--- a/doc/src/sgml/stream-check.sgml
+++ b/doc/src/sgml/stream-check.sgml
@@ -212,9 +212,10 @@
<listitem>
<para>
- Specifies the maximum tolerance level of replication delay in
- seconds on the standby server against the primary server. If the
- specified value is greater than
+ Specifies the maximum tolerance level of replication delay
+ on the standby server against the primary server.
+ If this value is specified without units, it is taken as milliseconds.
+ If the specified value is greater than
0, <xref linkend="guc-delay-threshold"> is ignored. If the delay
exceeds this configured level,
<productname>Pgpool-II</productname> stops sending the <acronym>
diff --git a/src/config/pool_config_variables.c b/src/config/pool_config_variables.c
index c5a6fb34b..639cf2250 100644
--- a/src/config/pool_config_variables.c
+++ b/src/config/pool_config_variables.c
@@ -2295,7 +2295,7 @@ static struct config_int ConfigureNamesInt[] =
{
{"delay_threshold_by_time", CFGCXT_RELOAD, STREAMING_REPLICATION_CONFIG,
"standby delay threshold by time.",
- CONFIG_VAR_TYPE_INT, false, GUC_UNIT_S,
+ CONFIG_VAR_TYPE_INT, false, GUC_UNIT_MS,
},
&g_pool_config.delay_threshold_by_time,
0,
diff --git a/src/context/pool_query_context.c b/src/context/pool_query_context.c
index 8e89a53be..50d15e33e 100644
--- a/src/context/pool_query_context.c
+++ b/src/context/pool_query_context.c
@@ -2018,8 +2018,6 @@ where_to_send_main_replica(POOL_QUERY_CONTEXT * query_context, char *query, Node
!pool_is_failed_transaction() &&
pool_get_transaction_isolation() != POOL_SERIALIZABLE))
{
- BackendInfo *bkinfo = pool_get_node_info(session_context->load_balance_node_id);
-
/*
* Load balance if possible
*/
@@ -2097,7 +2095,6 @@ where_to_send_main_replica(POOL_QUERY_CONTEXT * query_context, char *query, Node
if (pool_config->statement_level_load_balance)
{
session_context->load_balance_node_id = select_load_balancing_node();
- bkinfo = pool_get_node_info(session_context->load_balance_node_id);
}
/*
@@ -2106,12 +2103,7 @@ where_to_send_main_replica(POOL_QUERY_CONTEXT * query_context, char *query, Node
* load balance node which is lowest delayed,
* false then send to the primary.
*/
- if (STREAM &&
- (
- (pool_config->delay_threshold &&
- (bkinfo->standby_delay > pool_config->delay_threshold)) ||
- (pool_config->delay_threshold_by_time &&
- (bkinfo->standby_delay > pool_config->delay_threshold_by_time*1000*1000))))
+ if (STREAM && check_replication_delay(session_context->load_balance_node_id))
{
ereport(DEBUG1,
(errmsg("could not load balance because of too much replication delay"),
diff --git a/src/include/protocol/pool_pg_utils.h b/src/include/protocol/pool_pg_utils.h
index 25de6b2a6..bd9493572 100644
--- a/src/include/protocol/pool_pg_utils.h
+++ b/src/include/protocol/pool_pg_utils.h
@@ -60,5 +60,6 @@ extern void si_acquire_snapshot(void);
extern void si_snapshot_acquired(void);
extern void si_commit_request(void);
extern void si_commit_done(void);
+extern int check_replication_delay(int node_id);
#endif /* pool_pg_utils_h */
diff --git a/src/protocol/pool_pg_utils.c b/src/protocol/pool_pg_utils.c
index a81dfd170..8faff3e8b 100644
--- a/src/protocol/pool_pg_utils.c
+++ b/src/protocol/pool_pg_utils.c
@@ -441,10 +441,9 @@ select_load_balancing_node(void)
* and prefer_lower_delay_standby are true, we choose the least delayed
* node if suggested_node is standby and delayed over delay_threshold.
*/
- if (STREAM && pool_config->prefer_lower_delay_standby && suggested_node_id != PRIMARY_NODE_ID &&
- ((BACKEND_INFO(suggested_node_id).standby_delay_by_time && BACKEND_INFO(suggested_node_id).standby_delay > pool_config->delay_threshold_by_time * 1000000) ||
- (BACKEND_INFO(suggested_node_id).standby_delay_by_time == false && BACKEND_INFO(suggested_node_id).standby_delay > pool_config->delay_threshold)))
-
+ if (STREAM && pool_config->prefer_lower_delay_standby &&
+ suggested_node_id != PRIMARY_NODE_ID &&
+ check_replication_delay(suggested_node_id) < 0)
{
ereport(DEBUG1,
(errmsg("selecting load balance node"),
@@ -455,7 +454,7 @@ select_load_balancing_node(void)
* nodes which have the lowest delay.
*/
if (pool_config->delay_threshold_by_time > 0)
- lowest_delay = pool_config->delay_threshold_by_time * 1000 * 1000;
+ lowest_delay = pool_config->delay_threshold_by_time * 1000; /* convert from milli seconds to micro seconds */
else
lowest_delay = pool_config->delay_threshold;
@@ -602,17 +601,14 @@ select_load_balancing_node(void)
* node if suggested_node is standby and delayed over delay_threshold.
*/
if (STREAM && pool_config->prefer_lower_delay_standby &&
- ((pool_config->delay_threshold_by_time &&
- BACKEND_INFO(selected_slot).standby_delay > pool_config->delay_threshold_by_time*1000*1000) ||
- (pool_config->delay_threshold &&
- BACKEND_INFO(selected_slot).standby_delay > pool_config->delay_threshold)))
+ check_replication_delay(selected_slot) < 0)
{
ereport(DEBUG1,
(errmsg("selecting load balance node"),
errdetail("backend id %d is streaming delayed over delay_threshold", selected_slot)));
if (pool_config->delay_threshold_by_time > 0)
- lowest_delay = pool_config->delay_threshold_by_time * 1000 * 1000;
+ lowest_delay = pool_config->delay_threshold_by_time * 1000;
else
lowest_delay = pool_config->delay_threshold;
total_weight = 0.0;
@@ -1097,3 +1093,40 @@ si_commit_done(void)
session->si_state = SI_NO_SNAPSHOT;
}
}
+
+/*
+ * Check replication delay and returns the status.
+ * Return values:
+ * 0: no delay or not in streaming repplication mode or
+ * delay_threshold(_by_time) is set to 0
+ * -1: delay exceeds delay_threshold_by_time
+ * -2: delay exceeds delay_threshold
+ */
+int check_replication_delay(int node_id)
+{
+ BackendInfo *bkinfo;
+
+ if (!STREAM)
+ return 0;
+
+ bkinfo = pool_get_node_info(node_id);
+
+ /*
+ * Check delay_threshold_by_time. bkinfo->standby_delay is in
+ * microseconds while delay_threshold_by_time is in milliseconds. We need
+ * to multiply delay_threshold_by_time by 1000 to normalize.
+ */
+ if (pool_config->delay_threshold_by_time > 0 &&
+ bkinfo->standby_delay > pool_config->delay_threshold_by_time*1000)
+ return -1;
+
+ /*
+ * Check delay_threshold.
+ */
+ if (pool_config->delay_threshold > 0 &&
+ bkinfo->standby_delay > pool_config->delay_threshold)
+ return -2;
+
+ return 0;
+}
+
diff --git a/src/sample/pgpool.conf.sample-stream b/src/sample/pgpool.conf.sample-stream
index 769c516ca..e7b5b1c53 100644
--- a/src/sample/pgpool.conf.sample-stream
+++ b/src/sample/pgpool.conf.sample-stream
@@ -520,7 +520,7 @@ backend_clustering_mode = 'streaming_replication'
# Disabled (0) by default
#delay_threshold_by_time = 0
# Threshold before not dispatching query to standby node
- # Unit is in second(s)
+ # The default unit is in millisecond(s)
# Disabled (0) by default
#prefer_lower_delay_standby = off
@@ -679,7 +679,7 @@ backend_clustering_mode = 'streaming_replication'
#auto_failback = off
# Dettached backend node reattach automatically
- # if replication_state is 'streaming'.
+ # if replicatiotate is 'streaming'.
#auto_failback_interval = 1min
# Min interval of executing auto_failback in
# seconds.
diff --git a/src/streaming_replication/pool_worker_child.c b/src/streaming_replication/pool_worker_child.c
index 31bc1a62a..7b69dd7b2 100644
--- a/src/streaming_replication/pool_worker_child.c
+++ b/src/streaming_replication/pool_worker_child.c
@@ -495,7 +495,7 @@ check_replication_time_lag(void)
{
bkinfo->standby_delay = atol(s);
ereport(DEBUG1,
- (errmsg("standby delay in seconds * 1000000: " UINT64_FORMAT "", bkinfo->standby_delay)));
+ (errmsg("standby delay in milli seconds * 1000: " UINT64_FORMAT "", bkinfo->standby_delay)));
}
else
bkinfo->standby_delay = 0;
@@ -545,7 +545,7 @@ check_replication_time_lag(void)
{
lag = bkinfo->standby_delay;
delay_threshold_by_time = pool_config->delay_threshold_by_time;
- delay_threshold_by_time *= 1000000;
+ delay_threshold_by_time *= 1000; /* convert from milli seconds to micro seconds */
/* Log delay if necessary */
if ((pool_config->log_standby_delay == LSD_ALWAYS && lag > 0) ||
diff --git a/src/test/regression/tests/033.prefer_lower_standby_delay/test.sh b/src/test/regression/tests/033.prefer_lower_standby_delay/test.sh
index 9dc437693..d877af6a4 100755
--- a/src/test/regression/tests/033.prefer_lower_standby_delay/test.sh
+++ b/src/test/regression/tests/033.prefer_lower_standby_delay/test.sh
@@ -90,6 +90,7 @@ echo "delay_threshold = 10" >> etc/pgpool.conf
echo "sr_check_period = 1" >> etc/pgpool.conf
echo "log_standby_delay = 'always'" >> etc/pgpool.conf
echo "log_min_messages = 'DEBUG1'" >> etc/pgpool.conf
+echo "log_error_verbosity = verbose" >> etc/pgpool.conf
# force load balance node to be 1.
echo "backend_weight0 = 0" >> etc/pgpool.conf
echo "backend_weight2 = 0" >> etc/pgpool.conf
@@ -130,7 +131,8 @@ echo === Test2: delay_threshold_by_time with prefer_lower_delay_standby disabled
# ----------------------------------------------------------------------------------------
echo Start testing delay_threshold_by_time with prefer_lower_delay_standby disabled
echo "delay_threshold = 0" >> etc/pgpool.conf
-echo "delay_threshold_by_time = 1" >> etc/pgpool.conf
+echo "delay_threshold_by_time = 1000" >> etc/pgpool.conf
+
./startall
wait_for_pgpool_startup
# pause replay on node 1
[parent: 8f760f06d311]