agora inbox for pgsql-hackers@postgresql.orghelp / color / mirror / Atom feed
[PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. 266+ messages / 2 participants [nested] [flat]
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH] Collapse consecutive .** accessors for jsonpath exists queries @ 2026-06-18 20:30 Andrey Rachitskiy <pl0h0yp1@gmail.com> 0 siblings, 0 replies; 266+ messages in thread From: Andrey Rachitskiy @ 2026-06-18 20:30 UTC (permalink / raw) When a jsonpath expression contains multiple consecutive .** (jpiAny) accessors, each one triggers a full subtree traversal in executeAnyItem(). k consecutive .** operators can degrade performance to O(N^k) on a document with N nodes, even though a chain like $.**.**.** is redundant for existence semantics and equivalent to a single $.** with merged level bounds. This was reported as a performance problem when expressions such as $.**.**.**.**.* are evaluated with @? against deeply nested JSON: the same query without the trailing .* completes in sub-millisecond time, while the form with redundant .** segments can run for many minutes. A similar pattern in strict mode can also provoke very large memory allocation attempts. Collapse consecutive jpiAny nodes at execution time for existence queries (@? and jsonb_path_exists), merging their level bounds and performing a single executeAnyItem() pass. Author: Andrey Rachitskiy <pl0h0yp1@gmail.com> Reported-by: Andrey Rachitskiy <pl0h0yp1@gmail.com> Backpatch-through: 15 --- src/backend/utils/adt/jsonpath_exec.c | 47 +++++++++++++++++++--- .../src/test/regress/expected/jsonb_jsonpath.out | 40 ++++++++++++++++++ src/test/regress/sql/jsonb_jsonpath.sql | 11 +++++ 3 files changed, 93 insertions(+), 5 deletions(-) diff --git a/src/backend/utils/adt/jsonpath_exec.c b/src/backend/utils/adt/jsonpath_exec.c index 6cc2acb..7dfc311 100644 --- a/src/backend/utils/adt/jsonpath_exec.c +++ b/src/backend/utils/adt/jsonpath_exec.c @@ -113,6 +113,8 @@ typedef struct JsonPathExecContext * ignored */ bool throwErrors; /* with "false" all suppressible errors are * suppressed */ + bool existsOnly; /* @? / jsonb_path_exists: may collapse + * redundant consecutive .** accessors */ bool useTz; } JsonPathExecContext; @@ -310,6 +312,7 @@ static JsonPathExecResult executeAnyItem(JsonPathExecContext *cxt, JsonPathItem *jsp, JsonbContainer *jbc, JsonValueList *found, uint32 level, uint32 first, uint32 last, bool ignoreStructuralErrors, bool unwrapNext); +static uint32 mergeAnyBound(uint32 a, uint32 b); static JsonPathBool executePredicate(JsonPathExecContext *cxt, JsonPathItem *pred, JsonPathItem *larg, JsonPathItem *rarg, JsonbValue *jb, bool unwrapRightArg, @@ -739,6 +742,7 @@ executeJsonPath(JsonPath *path, void *vars, JsonPathGetVarCallback getVar, cxt.lastGeneratedObjectId = 1 + countVars(vars); cxt.innermostArraySize = -1; cxt.throwErrors = throwErrors; + cxt.existsOnly = (result == NULL); cxt.useTz = useTz; if (jspStrictAbsenceOfErrors(&cxt) && !result) @@ -1017,16 +1021,37 @@ executeItemOptUnwrapTarget(JsonPathExecContext *cxt, JsonPathItem *jsp, case jpiAny: { - bool hasNext = jspGetNext(jsp, &elem); + JsonPathItem elem; + bool hasNext; + uint32 first; + uint32 last; + + hasNext = jspGetNext(jsp, &elem); + first = jsp->content.anybounds.first; + last = jsp->content.anybounds.last; + + /* + * Consecutive .** accessors are redundant for existence + * queries and multiply traversal cost. + */ + if (cxt->existsOnly) + { + while (hasNext && elem.type == jpiAny) + { + first = mergeAnyBound(first, elem.content.anybounds.first); + last = mergeAnyBound(last, elem.content.anybounds.last); + hasNext = jspGetNext(&elem, &elem); + } + } /* first try without any intermediate steps */ - if (jsp->content.anybounds.first == 0) + if (first == 0) { bool savedIgnoreStructuralErrors; savedIgnoreStructuralErrors = cxt->ignoreStructuralErrors; cxt->ignoreStructuralErrors = true; - res = executeNextItem(cxt, jsp, &elem, + res = executeNextItem(cxt, jsp, hasNext ? &elem : NULL, jb, found); cxt->ignoreStructuralErrors = savedIgnoreStructuralErrors; @@ -1039,8 +1064,8 @@ executeItemOptUnwrapTarget(JsonPathExecContext *cxt, JsonPathItem *jsp, (cxt, hasNext ? &elem : NULL, jb->val.binary.data, found, 1, - jsp->content.anybounds.first, - jsp->content.anybounds.last, + first, + last, true, jspAutoUnwrap(cxt)); break; } @@ -1978,6 +2003,18 @@ executeNestedBoolItem(JsonPathExecContext *cxt, JsonPathItem *jsp, return res; } + +/* + * Merge level bounds of two consecutive .** accessors. + */ +static uint32 +mergeAnyBound(uint32 a, uint32 b) +{ + if (a == PG_UINT32_MAX || b == PG_UINT32_MAX) + return PG_UINT32_MAX; + return a + b; +} + /* * Implementation of several jsonpath nodes: * - jpiAny (.** accessor), diff --git a/src/test/regress/expected/jsonb_jsonpath.out b/src/test/regress/expected/jsonb_jsonpath.out index 81efebc..21f1362 100644 --- a/src/test/regress/expected/jsonb_jsonpath.out +++ b/src/test/regress/expected/jsonb_jsonpath.out @@ -785,6 +785,46 @@ select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{2 to 3}.b ? ( @ > 0)'; t (1 row) +-- Redundant consecutive .** accessors are collapsed for existence queries +-- (@? and jsonb_path_exists), avoiding multiplicative traversal cost. +select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.b ? (@ > 0)'; + ?column? +---------- + t +(1 row) + +select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.**.b ? (@ > 0)'; + ?column? +---------- + t +(1 row) + +select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.**.b ? (@ > 99)'; + ?column? +---------- + f +(1 row) + +select jsonb_path_exists('{"a": {"c": {"b": 1}}}', '$.**.**.b ? (@ > 0)'); + jsonb_path_exists +------------------- + t +(1 row) + +select ('[' || repeat('[', 50) || '0' || repeat(']', 50) || ']')::jsonb + @? 'lax $.**.**.**.**'; + ?column? +---------- + t +(1 row) + +select ('[' || repeat('[', 50) || '0' || repeat(']', 50) || ']')::jsonb + @? 'strict $.**.**.**'; + ?column? +---------- + t +(1 row) + select jsonb_path_query('{"g": {"x": 2}}', '$.g ? (exists (@.x))'); jsonb_path_query ------------------ diff --git a/src/test/regress/sql/jsonb_jsonpath.sql b/src/test/regress/sql/jsonb_jsonpath.sql index c1f4ab5..92abe13 100644 --- a/src/test/regress/sql/jsonb_jsonpath.sql +++ b/src/test/regress/sql/jsonb_jsonpath.sql @@ -150,6 +150,17 @@ select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{0 to last}.b ? ( @ > 0)'; select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{1 to last}.b ? ( @ > 0)'; select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{1 to 2}.b ? ( @ > 0)'; select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**{2 to 3}.b ? ( @ > 0)'; +-- Redundant consecutive .** accessors are collapsed for existence queries +-- (@? and jsonb_path_exists), avoiding multiplicative traversal cost. +select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.b ? (@ > 0)'; +select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.**.b ? (@ > 0)'; +select jsonb '{"a": {"c": {"b": 1}}}' @? '$.**.**.b ? (@ > 99)'; +select jsonb_path_exists('{"a": {"c": {"b": 1}}}', '$.**.**.b ? (@ > 0)'); +select ('[' || repeat('[', 50) || '0' || repeat(']', 50) || ']')::jsonb + @? 'lax $.**.**.**.**'; +select ('[' || repeat('[', 50) || '0' || repeat(']', 50) || ']')::jsonb + @? 'strict $.**.**.**'; + select jsonb_path_query('{"g": {"x": 2}}', '$.g ? (exists (@.x))'); select jsonb_path_query('{"g": {"x": 2}}', '$.g ? (exists (@.y))'); -- 2.47.3 --MP_/BlgiDE/t55y8Vb7V2uvNybc Content-Type: text/x-patch Content-Transfer-Encoding: 7bit Content-Disposition: attachment; filename=0001-collapse-consecutive-jsonpath-any-pg15.patch ^ permalink raw reply [nested|flat] 266+ messages in thread
end of thread, other threads:[~2026-06-18 20:30 UTC | newest] Thread overview: 266+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <andres@anarazel.de> 2026-06-18 20:30 [PATCH] Collapse consecutive .** accessors for jsonpath exists queries Andrey Rachitskiy <pl0h0yp1@gmail.com>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox