agora inbox for pgsql-hackers@postgresql.org  
help / color / mirror / Atom feed
[PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
266+ messages / 2 participants
[nested] [flat]

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation.
@ 2020-06-12 02:38 Andres Freund <andres@anarazel.de>
  0 siblings, 0 replies; 266+ messages in thread

From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw)

---
 src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++----
 1 file changed, 60 insertions(+), 8 deletions(-)

diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h
index fc058d548a8..8b2f9a2e707 100644
--- a/src/include/portability/instr_time.h
+++ b/src/include/portability/instr_time.h
@@ -83,7 +83,9 @@
 #define PG_INSTR_CLOCK	CLOCK_REALTIME
 #endif
 
+/* time in baseline cpu cycles */
 typedef int64 instr_time;
+
 #define NS_PER_S INT64CONST(1000000000)
 #define US_PER_S INT64CONST(1000000)
 #define MS_PER_S INT64CONST(1000)
@@ -95,17 +97,67 @@ typedef int64 instr_time;
 
 #define INSTR_TIME_SET_ZERO(t)	((t) = 0)
 
-static inline instr_time pg_clock_gettime_ns(void)
+#include <x86intrin.h>
+#include <cpuid.h>
+
+/*
+ * Return what the number of cycles needs to be multiplied with to end up with
+ * seconds.
+ *
+ * FIXME: The cold portion should probably be out-of-line. And it'd be better
+ * to not recompute this in every file that uses this. Best would probably be
+ * to require explicit initialization of cycles_to_sec, because having a
+ * branch really is unnecessary.
+ *
+ * FIXME: We should probably not unnecessarily use floating point math
+ * here. And it's likely that the numbers are small enough that we are running
+ * into floating point inaccuracies already. Probably worthwhile to be a good
+ * bit smarter.
+ *
+ * FIXME: This would need to be conditional, with a fallback to something not
+ * rdtsc based.
+ */
+static inline double __attribute__((const))
+get_cycles_to_sec(void)
 {
-	struct timespec tmp;
+	static double cycles_to_sec = 0;
 
-	clock_gettime(PG_INSTR_CLOCK, &tmp);
+	/*
+	 * Compute baseline cpu peformance, determines speed at which rdtsc advances
+	 */
+	if (unlikely(cycles_to_sec == 0))
+	{
+		uint32 cpuinfo[4] = {0};
 
-	return tmp.tv_sec * NS_PER_S + tmp.tv_nsec;
+		__get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3);
+		cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000);
+	}
+
+	return cycles_to_sec;
+}
+
+static inline instr_time pg_clock_gettime_ref_cycles(void)
+{
+	/*
+	 * The rdtscp waits for all in-flight instructions to finish (but allows
+	 * later instructions to start concurrently). That's good for some timing
+	 * situations (when the time is supposed to cover all the work), but
+	 * terrible for others (when sub-parts of work are measured, because then
+	 * the pipeline stall due to the wait change the overall timing).
+	 */
+#if 0
+	unsigned int aux;
+	int64 tsc = __rdtscp(&aux);
+
+	return tsc;
+#else
+
+	return __rdtsc();
+#endif
 }
 
 #define INSTR_TIME_SET_CURRENT(t) \
-	(t) = pg_clock_gettime_ns()
+	(t) = pg_clock_gettime_ref_cycles()
 
 #define INSTR_TIME_ADD(x,y) \
 	do { \
@@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void)
 	} while (0)
 
 #define INSTR_TIME_GET_DOUBLE(t) \
-	((double) (t) / NS_PER_S)
+	((double) (t) * get_cycles_to_sec())
 
 #define INSTR_TIME_GET_MILLISEC(t) \
-	((double) (t) / NS_PER_MS)
+	((double) (t) * (get_cycles_to_sec() * MS_PER_S))
 
 #define INSTR_TIME_GET_MICROSEC(t) \
-	((double) (t) / NS_PER_US)
+	((double) (t) * (get_cycles_to_sec() * US_PER_S))
 
 #else							/* !HAVE_CLOCK_GETTIME */
 
-- 
2.25.0.114.g5b0ca878e0


--wn4ncs637ccpvbhb--





^ permalink  raw  reply  [nested|flat] 266+ messages in thread

* [PATCH] Collapse consecutive .** accessors for jsonpath exists queries
@ 2026-06-18 20:30 Andrey Rachitskiy <pl0h0yp1@gmail.com>
  0 siblings, 0 replies; 266+ messages in thread

From: Andrey Rachitskiy @ 2026-06-18 20:30 UTC (permalink / raw)

When a jsonpath expression contains multiple consecutive .** (jpiAny)
accessors, each one triggers a full subtree traversal in executeAnyItem().
k consecutive .** operators can degrade performance to O(N^k) on a document
with N nodes, even though a chain like $.**.**.** is redundant for existence
semantics and equivalent to a single $.** with merged level bounds.

This was reported as a performance problem when expressions such as
$.**.**.**.**.* are evaluated with @? against deeply nested JSON: the same
query without the trailing .* completes in sub-millisecond time, while the
form with redundant .** segments can run for many minutes.  A similar pattern
in strict mode can also provoke very large memory allocation attempts.

Collapse consecutive jpiAny nodes at execution time for existence queries
(@? and jsonb_path_exists), merging their level bounds and performing a
single executeAnyItem() pass.

Author: Andrey Rachitskiy <pl0h0yp1@gmail.com>
Reported-by: Andrey Rachitskiy <pl0h0yp1@gmail.com>
Backpatch-through: 15
---
 src/backend/utils/adt/jsonpath_exec.c     | 47 +++++++++++++++++++---
 .../src/test/regress/expected/jsonb_jsonpath.out   | 40 ++++++++++++++++++
 src/test/regress/sql/jsonb_jsonpath.sql   | 11 +++++
 3 files changed, 93 insertions(+), 5 deletions(-)

diff --git a/src/backend/utils/adt/jsonpath_exec.c b/src/backend/utils/adt/jsonpath_exec.c
index 6cc2acb..7dfc311 100644
--- a/src/backend/utils/adt/jsonpath_exec.c
+++ b/src/backend/utils/adt/jsonpath_exec.c
@@ -113,6 +113,8 @@ typedef struct JsonPathExecContext
 										 * ignored */
 	bool		throwErrors;	/* with "false" all suppressible errors are
 								 * suppressed */
+	bool		existsOnly;		/* @? / jsonb_path_exists: may collapse
+							 * redundant consecutive .** accessors */
 	bool		useTz;
 } JsonPathExecContext;
 
@@ -310,6 +312,7 @@ static JsonPathExecResult executeAnyItem(JsonPathExecContext *cxt,
 										 JsonPathItem *jsp, JsonbContainer *jbc, JsonValueList *found,
 										 uint32 level, uint32 first, uint32 last,
 										 bool ignoreStructuralErrors, bool unwrapNext);
+static uint32 mergeAnyBound(uint32 a, uint32 b);
 static JsonPathBool executePredicate(JsonPathExecContext *cxt,
 									 JsonPathItem *pred, JsonPathItem *larg, JsonPathItem *rarg,
 									 JsonbValue *jb, bool unwrapRightArg,
@@ -739,6 +742,7 @@ executeJsonPath(JsonPath *path, void *vars, JsonPathGetVarCallback getVar,
 	cxt.lastGeneratedObjectId = 1 + countVars(vars);
 	cxt.innermostArraySize = -1;
 	cxt.throwErrors = throwErrors;
+	cxt.existsOnly = (result == NULL);
 	cxt.useTz = useTz;
 
 	if (jspStrictAbsenceOfErrors(&cxt) && !result)
@@ -1017,16 +1021,37 @@ executeItemOptUnwrapTarget(JsonPathExecContext *cxt, JsonPathItem *jsp,
 
 		case jpiAny:
 			{
-				bool		hasNext = jspGetNext(jsp, &elem);
+				JsonPathItem elem;
+				bool		hasNext;
+				uint32		first;
+				uint32		last;
+
+				hasNext = jspGetNext(jsp, &elem);
+				first = jsp->content.anybounds.first;
+				last = jsp->content.anybounds.last;
+
+				/*
+				 * Consecutive .** accessors are redundant for existence
+				 * queries and multiply traversal cost.
+				 */
+				if (cxt->existsOnly)
+				{
+					while (hasNext && elem.type == jpiAny)
+					{
+						first = mergeAnyBound(first, elem.content.anybounds.first);
+						last = mergeAnyBound(last, elem.content.anybounds.last);
+						hasNext = jspGetNext(&elem, &elem);
+					}
+				}
 
 				/* first try without any intermediate steps */
-				if (jsp->content.anybounds.first == 0)
+				if (first == 0)
 				{
 					bool		savedIgnoreStructuralErrors;
 
 					savedIgnoreStructuralErrors = cxt->ignoreStructuralErrors;
 					cxt->ignoreStructuralErrors = true;
-					res = executeNextItem(cxt, jsp, &elem,
+					res = executeNextItem(cxt, jsp, hasNext ? &elem : NULL,
 										  jb, found);
 					cxt->ignoreStructuralErrors = savedIgnoreStructuralErrors;
 
@@ -1039,8 +1064,8 @@ executeItemOptUnwrapTarget(JsonPathExecContext *cxt, JsonPathItem *jsp,
 						(cxt, hasNext ? &elem : NULL,
 						 jb->val.binary.data, found,
 						 1,
-						 jsp->content.anybounds.first,
-						 jsp->content.anybounds.last,
+						 first,
+						 last,
 						 true, jspAutoUnwrap(cxt));
 				break;
 			}
@@ -1978,6 +2003,18 @@ executeNestedBoolItem(JsonPathExecContext *cxt, JsonPathItem *jsp,
 	return res;
 }
 
+
+/*
+ * Merge level bounds of two consecutive .** accessors.
+ */
+static uint32
+mergeAnyBound(uint32 a, uint32 b)
+{
+	if (a == PG_UINT32_MAX || b == PG_UINT32_MAX)
+		return PG_UINT32_MAX;
+	return a + b;
+}
+
 /*
  * Implementation of several jsonpath nodes:
  *  - jpiAny (.** accessor),
diff --git a/src/test/regress/expected/jsonb_jsonpath.out b/src/test/regress/expected/jsonb_jsonpath.out
index 81efebc..21f1362 100644
--- a/src/test/regress/expected/jsonb_jsonpath.out
+++ b/src/test/regress/expected/jsonb_jsonpath.out
@@ -785,6 +785,46 @@ select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{2 to 3}.b ? ( @ > 0)';
  t
 (1 row)
 
+-- Redundant consecutive .** accessors are collapsed for existence queries
+-- (@? and jsonb_path_exists), avoiding multiplicative traversal cost.
+select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.b ? (@ > 0)';
+ ?column? 
+----------
+ t
+(1 row)
+
+select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.**.b ? (@ > 0)';
+ ?column? 
+----------
+ t
+(1 row)
+
+select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.**.b ? (@ > 99)';
+ ?column? 
+----------
+ f
+(1 row)
+
+select jsonb_path_exists('{"a": {"c": {"b": 1}}}', '$.**.**.b ? (@ > 0)');
+ jsonb_path_exists 
+-------------------
+ t
+(1 row)
+
+select ('[' || repeat('[', 50) || '0' || repeat(']', 50) || ']')::jsonb
+	@? 'lax $.**.**.**.**';
+ ?column? 
+----------
+ t
+(1 row)
+
+select ('[' || repeat('[', 50) || '0' || repeat(']', 50) || ']')::jsonb
+	@? 'strict $.**.**.**';
+ ?column? 
+----------
+ t
+(1 row)
+
 select jsonb_path_query('{"g": {"x": 2}}', '$.g ? (exists (@.x))');
  jsonb_path_query 
 ------------------
diff --git a/src/test/regress/sql/jsonb_jsonpath.sql b/src/test/regress/sql/jsonb_jsonpath.sql
index c1f4ab5..92abe13 100644
--- a/src/test/regress/sql/jsonb_jsonpath.sql
+++ b/src/test/regress/sql/jsonb_jsonpath.sql
@@ -150,6 +150,17 @@ select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{0 to last}.b ? ( @ > 0)';
 select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{1 to last}.b ? ( @ > 0)';
 select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{1 to 2}.b ? ( @ > 0)';
 select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{2 to 3}.b ? ( @ > 0)';
+-- Redundant consecutive .** accessors are collapsed for existence queries
+-- (@? and jsonb_path_exists), avoiding multiplicative traversal cost.
+select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.b ? (@ > 0)';
+select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.**.b ? (@ > 0)';
+select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.**.b ? (@ > 99)';
+select jsonb_path_exists('{"a": {"c": {"b": 1}}}', '$.**.**.b ? (@ > 0)');
+select ('[' || repeat('[', 50) || '0' || repeat(']', 50) || ']')::jsonb
+	@? 'lax $.**.**.**.**';
+select ('[' || repeat('[', 50) || '0' || repeat(']', 50) || ']')::jsonb
+	@? 'strict $.**.**.**';
+
 
 select jsonb_path_query('{"g": {"x": 2}}', '$.g ? (exists (@.x))');
 select jsonb_path_query('{"g": {"x": 2}}', '$.g ? (exists (@.y))');
-- 
2.47.3

--MP_/BlgiDE/t55y8Vb7V2uvNybc
Content-Type: text/x-patch
Content-Transfer-Encoding: 7bit
Content-Disposition: attachment;
 filename=0001-collapse-consecutive-jsonpath-any-pg15.patch



^ permalink  raw  reply  [nested|flat] 266+ messages in thread


end of thread, other threads:[~2026-06-18 20:30 UTC | newest]

Thread overview: 266+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de>
2026-06-18 20:30 [PATCH] Collapse consecutive .** accessors for jsonpath exists queries Andrey Rachitskiy <pl0h0yp1@gmail.com>

This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox